exp_name="0302_qwen2.5_7B_math_grpo_fast1_${RANDOM}"
python mason.py \
    --cluster ai2/jupiter  \
    --workspace ai2/tulu-3-dev \
    --priority normal \
    --preemptible \
    --num_nodes 1 \
    --max_retries 0 \
    --budget ai2/oe-adapt \
    --gpus 8 -- source configs/beaker_configs/ray_node_setup.sh \&\& python open_instruct/grpo_vllm_thread_ray_gtrl.py \
    --exp_name $exp_name \
    --output_dir /weka/oe-adapt-default/costah/models/$exp_name \
    --dataset_mixer_list ai2-adapt-dev/rlvr_gsm8k_zs 1.0 \
    --dataset_mixer_list_splits train \
    --dataset_mixer_eval_list ai2-adapt-dev/rlvr_gsm8k_zs 1.0 \
    --dataset_mixer_eval_list_splits train \
    --max_token_length 2048 \
    --max_prompt_token_length 2048 \
    --response_length 2048 \
    --number_samples_per_prompt 4 \
    --model_name_or_path meta-llama/Llama-3.1-8B \
    --stop_strings "</answer>" \
    --add_r1_style_format_reward \
    --non_stop_penalty False \
    --stop_token eos \
    --penalty_reward_value 0.0 \
    --temperature 0.7 \
    --ground_truths_key ground_truth \
    --chat_template_name r1_simple_chat_postpend_think \
    --sft_messages_key messages \
    --learning_rate 3e-7 \
    --total_episodes 1000000 \
    --deepspeed_stage 3 \
    --per_device_train_batch_size 1 \
    --local_rollout_forward_batch_size 1 \
    --local_mini_batch_size 16 \
    --local_rollout_batch_size 16 \
    --num_epochs 1 \
    --actor_num_gpus_per_node 6 \
    --vllm_tensor_parallel_size 2 \
    --beta 0.01 \
    --apply_verifiable_reward true \
    --seed 3 \
    --num_evals 100 \
    --save_freq 100 \
    --reward_model_multiplier 0.0 \
    --no_try_launch_beaker_eval_jobs \
    --gradient_checkpointing \
    --with_tracking
